agora inbox for pgsql-hackers@postgresql.org  
help / color / mirror / Atom feed
[PATCH 5/6] Use only xlogreader.c:XLogRead()
402+ messages / 1 participants
[nested] [flat]

* [PATCH 5/6] Use only xlogreader.c:XLogRead()
@ 2019-09-23 05:40  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2019-09-23 05:40 UTC (permalink / raw)

The implementations in xlogutils.c and walsender.c are just renamed now, to be
removed by the following diff.
---
 src/backend/access/transam/xlogreader.c | 157 ++++++++++++++++++++++++
 src/backend/access/transam/xlogutils.c  |  46 ++++++-
 src/backend/replication/walsender.c     | 139 ++++++++++++++++++++-
 src/bin/pg_waldump/pg_waldump.c         |  64 +++++++++-
 src/include/access/xlogreader.h         |  42 +++++++
 5 files changed, 436 insertions(+), 12 deletions(-)

diff --git a/src/backend/access/transam/xlogreader.c b/src/backend/access/transam/xlogreader.c
index 4de5530b3e..7f77fa95cb 100644
--- a/src/backend/access/transam/xlogreader.c
+++ b/src/backend/access/transam/xlogreader.c
@@ -17,6 +17,8 @@
  */
 #include "postgres.h"
 
+#include <unistd.h>
+
 #include "access/transam.h"
 #include "access/xlogrecord.h"
 #include "access/xlog_internal.h"
@@ -27,6 +29,7 @@
 
 #ifndef FRONTEND
 #include "miscadmin.h"
+#include "pgstat.h"
 #include "utils/memutils.h"
 #endif
 
@@ -1015,6 +1018,160 @@ WALOpenSegmentInit(WALOpenSegment *seg, int size)
 #endif
 }
 
+/*
+ * Read 'count' bytes from WAL into 'buf', starting at location 'startptr'. If
+ * tli_p is passed, get the data from timeline *tli_p. 'pos' is the current
+ * position in the XLOG file and openSegment is a callback that opens the next
+ * segment for reading.
+ *
+ * Returns error information if the data could not be read or NULL if
+ * succeeded.
+ *
+ * XXX probably this should be improved to suck data directly from the
+ * WAL buffers when possible.
+ */
+XLogReadError *
+XLogRead(char *buf, XLogRecPtr startptr, Size count,
+		 TimeLineID *tli_p, WALOpenSegment *seg, WALSegmentOpen openSegment)
+{
+	char	   *p;
+	XLogRecPtr	recptr;
+	Size		nbytes;
+
+	p = buf;
+	recptr = startptr;
+	nbytes = count;
+
+	while (nbytes > 0)
+	{
+		int			segbytes;
+		int			readbytes;
+
+		seg->off = XLogSegmentOffset(recptr, seg->size);
+
+		if (seg->file < 0 ||
+			!XLByteInSeg(recptr, seg->num, seg->size) ||
+			(tli_p != NULL && *tli_p != seg->tli))
+		{
+			XLogSegNo	nextSegNo;
+			TimeLineID	tli = InvalidTimeLineID;
+			int			file;
+
+			/* Switch to another logfile segment */
+			if (seg->file >= 0)
+				close(seg->file);
+
+			XLByteToSeg(recptr, nextSegNo, seg->size);
+
+			/* If we have the TLI, let's pass it to the callback. */
+			if (tli_p != NULL)
+				tli = *tli_p;
+
+			/* Open the next segment in the caller's way. */
+			openSegment(nextSegNo, &tli, &file, seg);
+
+			/*
+			 * If we passed InvalidTimeLineID, the callback should have
+			 * determined the correct TLI and returned it.
+			 */
+			Assert(tli != InvalidTimeLineID);
+
+			/* Update the open segment info. */
+			seg->tli = tli;
+			seg->file = file;
+
+			/*
+			 * If the function is called by the XLOG reader, the reader will
+			 * eventually set both "num" and "off". However we need to care
+			 * about them too because the function can also be used directly,
+			 * see walsender.c.
+			 */
+			seg->num = nextSegNo;
+			seg->off = 0;
+		}
+
+		/* How many bytes are within this segment? */
+		if (nbytes > (seg->size - seg->off))
+			segbytes = seg->size - seg->off;
+		else
+			segbytes = nbytes;
+
+#ifndef FRONTEND
+		pgstat_report_wait_start(WAIT_EVENT_WAL_READ);
+#endif
+
+		/*
+		 * Failure to read the data does not necessarily imply non-zero errno.
+		 * Set it to zero so that caller can distinguish the failure that does
+		 * not affect errno.
+		 */
+		errno = 0;
+
+		readbytes = pg_pread(seg->file, p, segbytes, seg->off);
+
+#ifndef FRONTEND
+		pgstat_report_wait_end();
+#endif
+
+		if (readbytes <= 0)
+		{
+			XLogReadError *errinfo;
+
+			errinfo = (XLogReadError *) palloc(sizeof(XLogReadError));
+			errinfo->read_errno = errno;
+			errinfo->readbytes = readbytes;
+			errinfo->reqbytes = segbytes;
+			errinfo->seg = seg;
+
+			return errinfo;
+		}
+
+		/* Update state for read */
+		recptr += readbytes;
+		nbytes -= readbytes;
+		p += readbytes;
+
+		/*
+		 * If the function is called by the XLOG reader, the reader will
+		 * eventually set this field. However we need to care about it too
+		 * because the function can also be used directly (see walsender.c).
+		 */
+		seg->off += readbytes;
+	}
+
+	return NULL;
+}
+
+#ifndef FRONTEND
+/*
+ * Backend-specific convenience code to handle read errors encountered by
+ * XLogRead().
+ */
+void
+XLogReadProcessError(XLogReadError *errinfo)
+{
+	WALOpenSegment *seg = errinfo->seg;
+
+	if (errinfo->readbytes < 0)
+	{
+		errno = errinfo->read_errno;
+		ereport(ERROR,
+				(errcode_for_file_access(),
+				 errmsg("could not read from log segment %s, offset %u, length %zu: %m",
+						XLogFileNameP(seg->tli, seg->num), seg->off,
+						(Size) errinfo->reqbytes)));
+	}
+	else
+	{
+		ereport(ERROR,
+				(errcode(ERRCODE_DATA_CORRUPTED),
+				 errmsg("could not read from log segment %s, offset %u: length %zu",
+						XLogFileNameP(seg->tli, seg->num), seg->off,
+						(Size) errinfo->reqbytes)));
+	}
+}
+#endif
+
 /* ----------------------------------------
  * Functions for decoding the data and block references in a record.
  * ----------------------------------------
diff --git a/src/backend/access/transam/xlogutils.c b/src/backend/access/transam/xlogutils.c
index 424bb06919..38c3196168 100644
--- a/src/backend/access/transam/xlogutils.c
+++ b/src/backend/access/transam/xlogutils.c
@@ -653,8 +653,8 @@ XLogTruncateRelation(RelFileNode rnode, ForkNumber forkNum,
  * frontend).  Probably these should be merged at some point.
  */
 static void
-XLogRead(char *buf, int segsize, TimeLineID tli, XLogRecPtr startptr,
-		 Size count)
+XLogReadOld(char *buf, int segsize, TimeLineID tli, XLogRecPtr startptr,
+			Size count)
 {
 	char	   *p;
 	XLogRecPtr	recptr;
@@ -896,6 +896,39 @@ XLogReadDetermineTimeline(XLogReaderState *state, XLogRecPtr wantPage, uint32 wa
 	}
 }
 
+/*
+ * Callback for XLogRead() to open the next segment.
+ */
+static void
+read_local_xlog_page_segment_open(XLogSegNo nextSegNo, TimeLineID *tli_p,
+								  int *file_p, WALOpenSegment *seg)
+{
+	TimeLineID	tli = *tli_p;
+	char		path[MAXPGPATH];
+	int			file;
+
+	Assert(tli != InvalidTimeLineID);
+
+	XLogFilePath(path, tli, nextSegNo, seg->size);
+	file = BasicOpenFile(path, O_RDONLY | PG_BINARY);
+
+	if (file < 0)
+	{
+		if (errno == ENOENT)
+			ereport(ERROR,
+					(errcode_for_file_access(),
+					 errmsg("requested WAL segment %s has already been removed",
+							path)));
+		else
+			ereport(ERROR,
+					(errcode_for_file_access(),
+					 errmsg("could not open file \"%s\": %m",
+							path)));
+	}
+
+	*file_p = file;
+}
+
 /*
  * read_page callback for reading local xlog files
  *
@@ -915,6 +948,7 @@ read_local_xlog_page(XLogReaderState *state, XLogRecPtr targetPagePtr,
 				loc;
 	int			count;
 	TimeLineID	pageTLI;
+	XLogReadError *errinfo;
 
 	loc = targetPagePtr + reqLen;
 
@@ -1022,10 +1056,10 @@ read_local_xlog_page(XLogReaderState *state, XLogRecPtr targetPagePtr,
 	 * as 'count', read the whole page anyway. It's guaranteed to be
 	 * zero-padded up to the page boundary if it's incomplete.
 	 */
-	XLogRead(cur_page, state->seg.size, state->seg.tli, targetPagePtr,
-			 XLOG_BLCKSZ);
-	state->seg.tli = pageTLI;
-
+	if ((errinfo = XLogRead(cur_page, targetPagePtr, XLOG_BLCKSZ, &pageTLI,
+							&state->seg,
+							read_local_xlog_page_segment_open)) != NULL)
+		XLogReadProcessError(errinfo);
 	/* number of valid bytes in the buffer */
 	return count;
 }
diff --git a/src/backend/replication/walsender.c b/src/backend/replication/walsender.c
index a617e20ab6..a7b3e0ecbe 100644
--- a/src/backend/replication/walsender.c
+++ b/src/backend/replication/walsender.c
@@ -247,9 +247,12 @@ static void LagTrackerWrite(XLogRecPtr lsn, TimestampTz local_flush_time);
 static TimeOffset LagTrackerRead(int head, XLogRecPtr lsn, TimestampTz now);
 static bool TransactionIdInRecentPast(TransactionId xid, uint32 epoch);
 
-static void XLogRead(char *buf, XLogRecPtr startptr, Size count);
+static void WalSndSegmentOpen(XLogSegNo nextSegNo, TimeLineID *tli_p,
+							  int *file_p, WALOpenSegment *seg);
 
 
+static void XLogReadOld(char *buf, XLogRecPtr startptr, Size count);
+
 /* Initialize walsender process before entering the main command loop */
 void
 InitWalSender(void)
@@ -763,6 +766,7 @@ logical_read_xlog_page(XLogReaderState *state, XLogRecPtr targetPagePtr, int req
 {
 	XLogRecPtr	flushptr;
 	int			count;
+	XLogReadError *errinfo;
 
 	XLogReadDetermineTimeline(state, targetPagePtr, reqLen);
 	sendTimeLineIsHistoric = (state->currTLI != ThisTimeLineID);
@@ -783,7 +787,13 @@ logical_read_xlog_page(XLogReaderState *state, XLogRecPtr targetPagePtr, int req
 		count = flushptr - targetPagePtr;	/* part of the page available */
 
 	/* now actually read the data, we know it's there */
-	XLogRead(cur_page, targetPagePtr, XLOG_BLCKSZ);
+	if ((errinfo = XLogRead(cur_page,
+							targetPagePtr,
+							XLOG_BLCKSZ,
+							NULL,	/* WalSndSegmentOpen will determine TLI */
+							sendSeg,
+							WalSndSegmentOpen)) != NULL)
+		XLogReadProcessError(errinfo);
 
 	return count;
 }
@@ -2360,7 +2370,7 @@ WalSndKill(int code, Datum arg)
  * more than one.
  */
 static void
-XLogRead(char *buf, XLogRecPtr startptr, Size count)
+XLogReadOld(char *buf, XLogRecPtr startptr, Size count)
 {
 	char	   *p;
 	XLogRecPtr	recptr;
@@ -2533,6 +2543,81 @@ retry:
 	}
 }
 
+/*
+ * Callback for XLogRead() to open the next segment.
+ */
+void
+WalSndSegmentOpen(XLogSegNo nextSegNo, TimeLineID *tli_p, int *file_p,
+				  WALOpenSegment *seg)
+{
+	TimeLineID	tli = *tli_p;
+	char		path[MAXPGPATH];
+	int			file;
+
+	/*
+	 * The timeline is determined below, caller should not pass it.
+	 */
+	Assert(tli == InvalidTimeLineID);
+
+	/*-------
+	 * When reading from a historic timeline, and there is a timeline switch
+	 * within this segment, read from the WAL segment belonging to the new
+	 * timeline.
+	 *
+	 * For example, imagine that this server is currently on timeline 5, and
+	 * we're streaming timeline 4. The switch from timeline 4 to 5 happened at
+	 * 0/13002088. In pg_wal, we have these files:
+	 *
+	 * ...
+	 * 000000040000000000000012
+	 * 000000040000000000000013
+	 * 000000050000000000000013
+	 * 000000050000000000000014
+	 * ...
+	 *
+	 * In this situation, when requested to send the WAL from segment 0x13, on
+	 * timeline 4, we read the WAL from file 000000050000000000000013. Archive
+	 * recovery prefers files from newer timelines, so if the segment was
+	 * restored from the archive on this server, the file belonging to the old
+	 * timeline, 000000040000000000000013, might not exist. Their contents are
+	 * equal up to the switchpoint, because at a timeline switch, the used
+	 * portion of the old segment is copied to the new file.  -------
+	 */
+	tli = sendTimeLine;
+	if (sendTimeLineIsHistoric)
+	{
+		XLogSegNo	endSegNo;
+
+		XLByteToSeg(sendTimeLineValidUpto, endSegNo, seg->size);
+		if (seg->num == endSegNo)
+			tli = sendTimeLineNextTLI;
+	}
+
+	XLogFilePath(path, tli, nextSegNo, seg->size);
+	file = BasicOpenFile(path, O_RDONLY | PG_BINARY);
+
+	if (file < 0)
+	{
+		/*
+		 * If the file is not found, assume it's because the standby asked for
+		 * a too old WAL segment that has already been removed or recycled.
+		 */
+		if (errno == ENOENT)
+			ereport(ERROR,
+					(errcode_for_file_access(),
+					 errmsg("requested WAL segment %s has already been removed",
+							XLogFileNameP(tli, nextSegNo))));
+		else
+			ereport(ERROR,
+					(errcode_for_file_access(),
+					 errmsg("could not open file \"%s\": %m",
+							path)));
+	}
+
+	*file_p = file;
+	*tli_p = tli;
+}
+
 /*
  * Send out the WAL in its normal physical/stored form.
  *
@@ -2550,6 +2635,8 @@ XLogSendPhysical(void)
 	XLogRecPtr	startptr;
 	XLogRecPtr	endptr;
 	Size		nbytes;
+	XLogSegNo	segno;
+	XLogReadError *errinfo;
 
 	/* If requested switch the WAL sender to the stopping state. */
 	if (got_STOPPING)
@@ -2765,7 +2852,51 @@ XLogSendPhysical(void)
 	 * calls.
 	 */
 	enlargeStringInfo(&output_message, nbytes);
-	XLogRead(&output_message.data[output_message.len], startptr, nbytes);
+
+retry:
+	if ((errinfo = XLogRead(&output_message.data[output_message.len],
+							startptr,
+							nbytes,
+							NULL,	/* WalSndSegmentOpen will determine TLI */
+							sendSeg,
+							WalSndSegmentOpen)) != NULL)
+		XLogReadProcessError(errinfo);
+
+	/*
+	 * After reading into the buffer, check that what we read was valid. We do
+	 * this after reading, because even though the segment was present when we
+	 * opened it, it might get recycled or removed while we read it. The
+	 * read() succeeds in that case, but the data we tried to read might
+	 * already have been overwritten with new WAL records.
+	 */
+	XLByteToSeg(startptr, segno, wal_segment_size);
+	CheckXLogRemoved(segno, ThisTimeLineID);
+
+	/*
+	 * During recovery, the currently-open WAL file might be replaced with the
+	 * file of the same name retrieved from archive. So we always need to
+	 * check what we read was valid after reading into the buffer. If it's
+	 * invalid, we try to open and read the file again.
+	 */
+	if (am_cascading_walsender)
+	{
+		WalSnd	   *walsnd = MyWalSnd;
+		bool		reload;
+
+		SpinLockAcquire(&walsnd->mutex);
+		reload = walsnd->needreload;
+		walsnd->needreload = false;
+		SpinLockRelease(&walsnd->mutex);
+
+		if (reload && sendSeg->file >= 0)
+		{
+			close(sendSeg->file);
+			sendSeg->file = -1;
+
+			goto retry;
+		}
+	}
+
 	output_message.len += nbytes;
 	output_message.data[output_message.len] = '\0';
 
diff --git a/src/bin/pg_waldump/pg_waldump.c b/src/bin/pg_waldump/pg_waldump.c
index a16793bb8b..cd5f589f03 100644
--- a/src/bin/pg_waldump/pg_waldump.c
+++ b/src/bin/pg_waldump/pg_waldump.c
@@ -296,6 +296,51 @@ identify_target_directory(XLogDumpPrivate *private, char *directory,
 		fatal_error("could not find any WAL file");
 }
 
+static void
+WALDumpOpenSegment(XLogSegNo nextSegNo, TimeLineID *tli_p, int *file_p,
+				   WALOpenSegment *seg)
+{
+	TimeLineID	tli = *tli_p;
+	char		fname[MAXPGPATH];
+	int			file;
+	int			tries;
+
+	Assert(tli != InvalidTimeLineID);
+
+	XLogFileName(fname, tli, nextSegNo, seg->size);
+
+	/*
+	 * In follow mode there is a short period of time after the server has
+	 * written the end of the previous file before the new file is available.
+	 * So we loop for 5 seconds looking for the file to appear before giving
+	 * up.
+	 */
+	for (tries = 0; tries < 10; tries++)
+	{
+		file = open_file_in_directory(seg->dir, fname);
+		if (file >= 0)
+			break;
+		if (errno == ENOENT)
+		{
+			int			save_errno = errno;
+
+			/* File not there yet, try again */
+			pg_usleep(500 * 1000);
+
+			errno = save_errno;
+			continue;
+		}
+		/* Any other error, fall through and fail */
+		break;
+	}
+
+	if (file < 0)
+		fatal_error("could not find file \"%s\": %s",
+					fname, strerror(errno));
+
+	*file_p = file;
+}
+
 /*
  * Read count bytes from a segment file in the specified directory, for the
  * given timeline, containing the specified record pointer; store the data in
@@ -427,6 +472,7 @@ XLogDumpReadPage(XLogReaderState *state, XLogRecPtr targetPagePtr, int reqLen,
 {
 	XLogDumpPrivate *private = state->private_data;
 	int			count = XLOG_BLCKSZ;
+	XLogReadError *errinfo;
 
 	if (private->endptr != InvalidXLogRecPtr)
 	{
@@ -441,8 +487,22 @@ XLogDumpReadPage(XLogReaderState *state, XLogRecPtr targetPagePtr, int reqLen,
 		}
 	}
 
-	XLogDumpXLogRead(private->inpath, private->timeline, targetPagePtr,
-					 readBuff, count);
+	if ((errinfo = XLogRead(readBuff, targetPagePtr, count, &private->timeline,
+							&state->seg, WALDumpOpenSegment)) != NULL)
+	{
+		WALOpenSegment *seg = errinfo->seg;
+		char		fname[MAXPGPATH];
+
+		XLogFileName(fname, seg->tli, seg->num, seg->size);
+
+		if (errno != 0)
+			fatal_error("could not read from log file %s, offset %u, length %zu: %s",
+						fname, seg->off, (Size) errinfo->reqbytes,
+						strerror(errinfo->read_errno));
+		else
+			fatal_error("could not read from log file %s, offset %u: length: %zu",
+						fname, seg->off, (Size) errinfo->reqbytes);
+	}
 
 	return count;
 }
diff --git a/src/include/access/xlogreader.h b/src/include/access/xlogreader.h
index b9d99d524e..3d9742b81b 100644
--- a/src/include/access/xlogreader.h
+++ b/src/include/access/xlogreader.h
@@ -227,8 +227,50 @@ extern bool XLogReaderValidatePageHeader(XLogReaderState *state,
 extern XLogRecPtr XLogFindNextRecord(XLogReaderState *state, XLogRecPtr RecPtr);
 #endif							/* FRONTEND */
 
+/*
+ * Callback to open the specified WAL segment for reading.
+ *
+ * "nextSegNo" is the number of the segment to be opened.
+ *
+ * "tli_p" is an input/output argument. If *tli_p is valid, it's the timeline
+ * the new segment should be in. If *tli_p==InvalidTimeLineID, the callback
+ * needs to determine the timeline itself and put the result into *tli_p.
+ *
+ * "file_p" points to an address the segment file descriptor should be stored
+ * at.
+ *
+ * "seg" provides information on the currently open segment. The callback is
+ * not supposed to change this info.
+ *
+ * BasicOpenFile() is the preferred way to open the segment file in backend
+ * code, whereas open(2) should be used in frontend.
+ */
+typedef void (*WALSegmentOpen) (XLogSegNo nextSegNo, TimeLineID *tli_p,
+								int *file_p, WALOpenSegment *seg);
+
 extern void WALOpenSegmentInit(WALOpenSegment *seg, int size);
 
+/*
+ * Error information that both backend and frontend caller can process.
+ *
+ * XXX Should the name be WALReadError? If so, we probably need to rename
+ * XLogRead() and XLogReadProcessError() too.
+ */
+typedef struct XLogReadError
+{
+	int			read_errno;		/* errno set by the last read(). */
+	int			readbytes;		/* Bytes read by the last read(). */
+	int			reqbytes;		/* Bytes requested to be read. */
+	WALOpenSegment *seg;		/* Segment we tried to read from. */
+} XLogReadError;
+
+extern XLogReadError *XLogRead(char *buf, XLogRecPtr startptr, Size count,
+							   TimeLineID *tli_p, WALOpenSegment *seg,
+							   WALSegmentOpen openSegment);
+#ifndef FRONTEND
+void		XLogReadProcessError(XLogReadError *errinfo);
+#endif
+
 /* Functions for decoding an XLogRecord */
 
 extern bool DecodeXLogRecord(XLogReaderState *state, XLogRecord *record,
-- 
2.20.1


--=-=-=
Content-Type: text/x-diff
Content-Disposition: attachment;
 filename=v06-0006-Remove-the-old-implemenations-of-XLogRead.patch



^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread

* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13  Antonin Houska <ah@cybertec.at>
  0 siblings, 0 replies; 402+ messages in thread

From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)

For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.

This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.

When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.

At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
 src/backend/commands/repack.c   |   4 +-
 src/backend/storage/lmgr/lmgr.c |  35 ++++++++---
 src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
 src/backend/storage/lmgr/proc.c |  32 +++++++---
 src/include/storage/lmgr.h      |   1 +
 src/include/storage/lock.h      |   5 +-
 src/include/storage/proc.h      |   4 +-
 7 files changed, 166 insertions(+), 22 deletions(-)

diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
 	/*
 	 * Acquire AccessExclusiveLock on the table, its TOAST relation (if there
 	 * is one), all its indexes, so that we can swap the files.
+	 *
+	 * TODO The same for indexes and TOAST?
 	 */
-	LockRelationOid(old_table_oid, AccessExclusiveLock);
+	LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
 
 	/*
 	 * Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
 	const ItemPointerData *ctid;
 } XactLockTableWaitInfo;
 
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+								  bool isUpgrade);
 static void XactLockTableWaitErrorCb(void *arg);
 
 /*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
  */
 void
 LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+	LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
 {
 	LOCKTAG		tag;
 	LOCALLOCK  *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, isUpgrade);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
 	SetLocktagRelationOid(&tag, relid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
 	SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
-							  false);
+							  false, false);
 
 	/*
 	 * Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
 						 relation->rd_lockInfo.lockRelId.relId);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
 					  ItemPointerGetOffsetNumber(tid));
 
 	return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
-								logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+								logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
 }
 
 /*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
 		SET_LOCKTAG_TRANSACTION(tag, xid);
 
 		if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
-								logLockFailure)
+								logLockFailure, false)
 			== LOCKACQUIRE_NOT_AVAIL)
 			return false;
 
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
 					   objsubid);
 
 	res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
-							  false);
+							  false, false);
 
 	if (res == LOCKACQUIRE_NOT_AVAIL)
 		return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
 			bool dontWait)
 {
 	return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
-							   true, NULL, false);
+							   true, NULL, false, false);
 }
 
 /*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
  *
  * logLockFailure indicates whether to log details when a lock acquisition
  * fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
  */
 LockAcquireResult
 LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 					bool dontWait,
 					bool reportMemoryError,
 					LOCALLOCK **locallockp,
-					bool logLockFailure)
+					bool logLockFailure,
+					bool isUpgrade)
 {
 	LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
 	LockMethod	lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		 * case, because JoinWaitQueue() may discover that we can acquire the
 		 * lock immediately after all.
 		 */
-		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+		waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+								   isUpgrade);
 	}
 
 	if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
 		Assert(!dontWait);
 		PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
 		LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+		/*
+		 * Lock upgrade can introduce deadlock, therefore enforce special
+		 * behavior of other processes that deal with this lock. In
+		 * particular, any waiter that sees upgradedBy set is expected to
+		 * perform deadlock check as soon as it's woken up. (Since we are
+		 * already in the queue, the deadlock detector has all the information
+		 * it needs.)
+		 */
+		if (isUpgrade)
+		{
+			dlist_iter	iter;
+
+			/*
+			 * There should not be multiple upgrades at the same time. XXX
+			 * ERROR ?
+			 */
+			Assert(lock->upgradedBy == NULL);
+
+			lock->upgradedBy = MyProc;
+
+			/*
+			 * Now, before I sleep myself, wake up all the existing waiters
+			 * (except for me), so they check for deadlock.
+			 */
+			dclist_foreach(iter, &lock->waitProcs)
+			{
+				PGPROC	   *proc = dlist_container(PGPROC, waitLink,
+												   iter.cur);
+
+				if (proc != MyProc)
+					SetLatch(&proc->procLatch);
+			}
+		}
 		LWLockRelease(partitionLock);
 
+		if (!isUpgrade)
+		{
+			HASH_SEQ_STATUS status;
+			LOCALLOCK	*llock;
+			bool	check_deadlock = false;
+
+			/*
+			 * Likewise, new waiters should check the isUpgrade flag and
+			 * engage deadlock detector if needed. The lock being acquired
+			 * should already be in the hash.
+			 *
+			 * Note that, to make deadlock detection happen as soon as
+			 * possible (i.e. regardless deadlock timeout), we check the other
+			 * locks too. XXX There might be ways to get into the deadlock
+			 * indirectly, but that needs more analysis. As long as the
+			 * upgrading process skips deadlock detection altogether, the
+			 * worst consequence of missing some lock wait cycle here is that
+			 * the other process will be kicked-off no sooner than after its
+			 * deadlock timeout has elapsed. (Which in turn means that the
+			 * lock upgrade will take longer than expected.)
+			 */
+			hash_seq_init(&status, LockMethodLocalHash);
+			while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+			{
+				LOCK	*mylock = llock->lock;
+
+				if (mylock && mylock->upgradedBy)
+				{
+					check_deadlock = true;
+					break;
+				}
+			}
+			/* Check for deadlock if needed. */
+			if (check_deadlock)
+			{
+				DeadLockState	deadlock_state;
+
+				deadlock_state = CheckDeadLock();
+				if (deadlock_state == DS_HARD_DEADLOCK)
+					DeadLockReport();
+			}
+		}
+
 		waitResult = WaitOnLock(locallock, owner);
 
 		/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
 			DeadLockReport();
 			/* DeadLockReport() will not return */
 		}
+
+		/*
+		 * If finishing the lock upgrade, we should not get into a deadlock
+		 * anymore, so let others know that they do not have to care either.
+		 */
+		if (isUpgrade)
+		{
+			LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+			Assert(lock->upgradedBy != NULL);
+			lock->upgradedBy = NULL;
+			LWLockRelease(partitionLock);
+		}
 	}
 	else
 		LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
 		lock->nGranted = 0;
 		MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
 		MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+		lock->upgradedBy = NULL;
 		LOCK_PRINT("LockAcquire: new", lock, lockmode);
 	}
 	else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
 			elog(PANIC, "proclock table corrupted");
 	}
 
+	/*
+	 * Was this backend upgrading the lock?
+	 */
+	if (lock->upgradedBy == MyProc)
+		lock->upgradedBy = NULL;
+
 	if (lock->nRequested == 0)
 	{
 		/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
 static void RemoveProcFromArray(int code, Datum arg);
 static void ProcKill(int code, Datum arg);
 static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
 
 
 /*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
  * NOTES: The process queue is now a priority queue for locking.
  */
 ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+			  bool isUpgrade)
 {
 	LOCKMODE	lockmode = locallock->tag.mode;
 	LOCK	   *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
 			/* Must he wait for me? */
 			if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
 			{
-				/* Must I wait for him ? */
-				if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+				/*
+				 * Must I wait for him? I don't want a deadlock during lock
+				 * upgrade - other processes should fail on it.
+				 */
+				if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+					!isUpgrade)
 				{
 					/*
 					 * Yes, so we have a deadlock.  Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
 			(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
 							 PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
 			ResetLatch(MyLatch);
-			/* check for deadlocks first, as that's probably log-worthy */
-			if (got_deadlock_timeout)
+			/*
+			 * Check for deadlocks first, as that's probably log-worthy.
+			 *
+			 * Do not wait for the timeout if the lock is being upgraded since
+			 * the risk of deadlock is higher now. However, do not check for
+			 * deadlock if the lock is being upgraded by this process - other
+			 * processes should take care.
+			 *
+			 * TODO Possible optimization: if this is the only lock of the
+			 * backend and if it did not have any weaker lock on the table so
+			 * far, it should be safe to skip the deadlock check. However, to
+			 * evaluate the situation, we need to take fast-path locks into
+			 * account.
+			 */
+			if ((got_deadlock_timeout || lock->upgradedBy) &&
+				lock->upgradedBy != MyProc)
 			{
 				deadlock_state = CheckDeadLock();
 				got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
  * not, just return.  If we have a real deadlock, remove ourselves from the
  * lock's wait queue.
  */
-static DeadLockState
+DeadLockState
 CheckDeadLock(void)
 {
 	int			i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
 
 /* Lock a relation */
 extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
 extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
 extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
 extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
  * nRequested -- total requested locks of all types.
  * granted -- count of each lock type currently granted on the lock.
  * nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
  *
  * Note: these counts count 1 for each backend.  Internally to a backend,
  * there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
 	int			nRequested;		/* total of requested[] array */
 	int			granted[MAX_LOCKMODES]; /* counts of granted locks */
 	int			nGranted;		/* total of granted[] array */
+	PGPROC		*upgradedBy;	/* is lock being upgraded by this process? */
 } LOCK;
 
 #define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
 											 bool dontWait,
 											 bool reportMemoryError,
 											 LOCALLOCK **locallockp,
-											 bool logLockFailure);
+											 bool logLockFailure,
+											 bool isUpgrade);
 extern void AbortStrongLockAcquire(void);
 extern void MarkLockClear(LOCALLOCK *locallock);
 extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
 extern void ProcReleaseLocks(bool isCommit);
 
 extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
-									LockMethod lockMethodTable, bool dontWait);
+									LockMethod lockMethodTable, bool dontWait,
+									bool isUpgrade);
 extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
 extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
 extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
 extern void CheckDeadLockAlert(void);
 extern void LockErrorCleanup(void);
 extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
-- 
2.47.3


--=-=-=--





^ permalink  raw  reply  [nested|flat] 402+ messages in thread


end of thread, other threads:[~2026-04-17 13:13 UTC | newest]

Thread overview: 402+ messages (download: mbox mbox.gz follow: Atom feed)
-- links below jump to the message on this page --
2019-09-23 05:40 [PATCH 5/6] Use only xlogreader.c:XLogRead() Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <ah@cybertec.at>

This inbox is served by agora; see mirroring instructions
for how to clone and mirror all data and code used for this inbox